# -*- coding: utf-8 -*-
"""
Web crawler for intangible‑heritage sports course data from the Ministry of Education Traditional Culture Inheritance Bases
Functions:
1. Read the excel file "清洗后_仅新增非遗体育主项目清单.xlsx" and extract cleaned‑up main‑item names as matching keywords
2. Fetch all three batches of the "Outstanding Traditional Chinese Culture Inheritance Bases in Regular Higher Education Institutions" list published by the Ministry of Education
3. Filter universities whose inheritance projects belong to traditional sports intangible heritage, and match against user keywords
4. Export search results to an excel file in unified format

Data sources:
- Batch 1 (2018): Document 教体艺厅函〔2018〕85号, 55 bases
- Batch 2 (2019): Document 教体艺厅函〔2019〕53号, 25 bases
- Batch 3 (2020): Document 教体艺厅函〔2020〕26号, 26 bases
Total 106 bases
"""

import pandas as pd
import time
import re
import os
import requests
from bs4 import BeautifulSoup

# ========================== File Path Configuration ==========================
INPUT_KEYWORDS = "清洗后_仅新增非遗体育主项目清单.xlsx"
OUTPUT_FILE = "检索结果5_教育部传统文化传承基地数据.xlsx"

# ========================== HTTP Request Header Configuration ==========================
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
    "Accept-Encoding": "gzip, deflate, br",
    "Connection": "keep-alive",
}

# ========================== Built‑in Base List (Full Three‑Batch Data) ==========================
# Batch 1 (2018): 55 bases
BATCH1_BASE = [
    # Traditional sports category (partially known)
    {"地区": "河北", "学校名称": "河北体育学院", "传承项目": "武术·八极拳"},
    {"地区": "河北", "学校名称": "邯郸学院", "传承项目": "太极"},
    {"地区": "上海", "学校名称": "上海体育学院", "传承项目": "武术"},
    {"地区": "浙江", "学校名称": "浙江大学", "传承项目": "龙舟"},
    {"地区": "山东", "学校名称": "曲阜师范大学", "传承项目": "孔子射艺"},
    {"地区": "河南", "学校名称": "河南大学", "传承项目": "太极拳"},
    {"地区": "四川", "学校名称": "内江师范学院", "传承项目": "峨眉武术"},
    {"地区": "辽宁", "学校名称": "渤海大学", "传承项目": "少北武术"},
    {"地区": "辽宁", "学校名称": "沈阳体育学院", "传承项目": "五行通背拳"},
    {"地区": "安徽", "学校名称": "亳州学院", "传承项目": "五禽戏"},
    # Non‑sports category (kept for reference, will not be matched)
    {"地区": "北京", "学校名称": "北京大学", "传承项目": "昆曲"},
    {"地区": "北京", "学校名称": "清华大学", "传承项目": "京昆"},
    {"地区": "北京", "学校名称": "北京师范大学", "传承项目": "中国话剧"},
    {"地区": "天津", "学校名称": "天津大学", "传承项目": "杨柳青年画"},
    {"地区": "天津", "学校名称": "天津师范大学", "传承项目": "相声"},
    {"地区": "江苏", "学校名称": "南京大学", "传承项目": "紫砂"},
    {"地区": "江苏", "学校名称": "东南大学", "传承项目": "古琴"},
    {"地区": "江苏", "学校名称": "南京航空航天大学", "传承项目": "剪纸"},
    {"地区": "江苏", "学校名称": "苏州大学", "传承项目": "江南丝竹"},
    {"地区": "江苏", "学校名称": "江苏师范大学", "传承项目": "五禽操"},
    {"地区": "江苏", "学校名称": "南京财经大学", "传承项目": "篆刻"},
    {"地区": "江苏", "学校名称": "江苏第二师范学院", "传承项目": "锡剧"},
    {"地区": "江苏", "学校名称": "苏州工艺美术职业技术学院", "传承项目": "苏绣"},
    {"地区": "江苏", "学校名称": "江南大学", "传承项目": "染织刺绣技艺"},
    {"地区": "江苏", "学校名称": "无锡工艺职业技术学院", "传承项目": "宜兴紫砂陶"},
    {"地区": "上海", "学校名称": "复旦大学", "传承项目": "吴越踏歌"},
    # ... remaining non‑sports items omitted; only known traditional sports items retained
]

# Batch 2 (2019): 25 bases
BATCH2_BASE = [
    # Traditional sports category
    {"地区": "上海", "学校名称": "上海对外经贸大学", "传承项目": "中华射艺"},
    {"地区": "山东", "学校名称": "山东体育学院", "传承项目": "螳螂拳"},
    # Non‑sports category
    {"地区": "北京", "学校名称": "北京理工大学", "传承项目": "传统手工艺术"},
    {"地区": "山西", "学校名称": "山西大学", "传承项目": "晋剧"},
    {"地区": "辽宁", "学校名称": "东北大学", "传承项目": "木偶戏"},
    {"地区": "吉林", "学校名称": "吉林艺术学院", "传承项目": "吉剧"},
    {"地区": "江苏", "学校名称": "南京财经大学", "传承项目": "篆刻"},
    {"地区": "江苏", "学校名称": "江苏第二师范学院", "传承项目": "锡剧"},
    {"地区": "江苏", "学校名称": "苏州工艺美术职业技术学院", "传承项目": "苏绣"},
    {"地区": "浙江", "学校名称": "丽水学院", "传承项目": "青瓷"},
    {"地区": "安徽", "学校名称": "滁州学院", "传承项目": "凤阳花鼓"},
    {"地区": "福建", "学校名称": "福建师范大学", "传承项目": "闽台地方戏曲"},
    {"地区": "江西", "学校名称": "赣南师范大学", "传承项目": "赣南采茶戏"},
    {"地区": "山东", "学校名称": "山东艺术学院", "传承项目": "山东年画"},
    {"地区": "湖北", "学校名称": "武汉大学", "传承项目": "楚剧"},
    {"地区": "湖北", "学校名称": "武汉纺织大学", "传承项目": "汉绣"},
    {"地区": "湖南", "学校名称": "湖南工艺美术职业学院", "传承项目": "湘绣"},
    {"地区": "广东", "学校名称": "华南师范大学", "传承项目": "岭南传统舞蹈"},
    {"地区": "广东", "学校名称": "深圳信息职业技术学院", "传承项目": "诗经文化"},
    {"地区": "四川", "学校名称": "成都纺织高等专科学校", "传承项目": "蜀绣"},
    {"地区": "贵州", "学校名称": "贵州民族大学", "传承项目": "侗族大歌"},
    {"地区": "云南", "学校名称": "云南民族大学", "传承项目": "云南扎染"},
    {"地区": "陕西", "学校名称": "西安建筑科技大学", "传承项目": "传统瓦作技艺"},
    {"地区": "陕西", "学校名称": "陕西师范大学", "传承项目": "陕西皮影"},
    {"地区": "陕西", "学校名称": "西安外事学院", "传承项目": "古琴"},
]

# Batch 3 (2020): 26 bases
BATCH3_BASE = [
    # Traditional sports category
    {"地区": "上海", "学校名称": "上海体育学院", "传承项目": "武术"},
    {"地区": "福建", "学校名称": "华侨大学", "传承项目": "舞狮（南狮）"},
    {"地区": "河南", "学校名称": "河南中医药大学", "传承项目": "八段锦"},
    {"地区": "新疆", "学校名称": "新疆师范大学", "传承项目": "新疆民族传统体育"},
    # Non‑sports category
    {"地区": "北京", "学校名称": "北京科技大学", "传承项目": "传统金属工艺"},
    {"地区": "北京", "学校名称": "北京舞蹈学院", "传承项目": "民族民间舞蹈"},
    {"地区": "天津", "学校名称": "南开大学", "传承项目": "中国书画"},
    {"地区": "河北", "学校名称": "燕山大学", "传承项目": "篆刻"},
    {"地区": "河北", "学校名称": "邢台学院", "传承项目": "邢窑白瓷"},
    {"地区": "内蒙古", "学校名称": "内蒙古艺术学院", "传承项目": "蒙古族传统音乐"},
    {"地区": "辽宁", "学校名称": "大连海事大学", "传承项目": "中华绳结"},
    {"地区": "吉林", "学校名称": "东北师范大学人文学院", "传承项目": "东北秧歌"},
    {"地区": "黑龙江", "学校名称": "黑龙江大学", "传承项目": "冰灯冰雕制作技艺"},
    {"地区": "上海", "学校名称": "华东师范大学", "传承项目": "篆刻"},
    {"地区": "江苏", "学校名称": "江南大学", "传承项目": "江南染织"},
    {"地区": "江苏", "学校名称": "无锡工艺职业技术学院", "传承项目": "宜兴紫砂陶"},
    {"地区": "浙江", "学校名称": "湖州师范学院", "传承项目": "蚕丝绸文化"},
    {"地区": "安徽", "学校名称": "中国科学技术大学", "传承项目": "文房四宝工艺"},
    {"地区": "山东", "学校名称": "临沂大学", "传承项目": "柳琴戏"},
    {"地区": "湖北", "学校名称": "中国地质大学（武汉）", "传承项目": "编钟艺术"},
    {"地区": "湖北", "学校名称": "黄冈师范学院", "传承项目": "黄梅戏"},
    {"地区": "广西", "学校名称": "广西大学", "传承项目": "彩调剧"},
    {"地区": "海南", "学校名称": "海南师范大学", "传承项目": "黎锦"},
    {"地区": "陕西", "学校名称": "榆林学院", "传承项目": "陕北秧歌"},
    {"地区": "陕西", "学校名称": "陕西艺术职业学院", "传承项目": "秦腔"},
    {"地区": "甘肃", "学校名称": "兰州城市学院", "传承项目": "西部民歌"},
]


# ========================== Data Reading Module ==========================
def read_keywords():
    """Read and return the cleaned‑up main‑item keyword list"""
    if not os.path.exists(INPUT_KEYWORDS):
        raise FileNotFoundError(f"Input file not found: {INPUT_KEYWORDS}")

    df = pd.read_excel(INPUT_KEYWORDS)
    if "清洗后主项目名称" not in df.columns:
        raise ValueError(f"Required column missing in excel. Current column list: {df.columns.tolist()}")

    keywords = df["清洗后主项目名称"].dropna().astype(str).str.strip().tolist()
    keywords = [k for k in keywords if k and k != "nan"]

    if not keywords:
        raise ValueError("No valid matching keywords extracted")

    print(f"✅ Successfully loaded {len(keywords)} matching keywords")
    return keywords


# ========================== Matching Logic ==========================
def match_keyword(project_name, keywords):
    """
    Determine whether the inheritance project matches the keyword list
    Return: the list of matched keywords (possibly multiple)
    """
    matched = []
    for kw in keywords:
        # Exact match or substring match (e.g., "太极拳" matches "太极")
        if kw == project_name:
            matched.append(kw)
        elif kw in project_name:
            matched.append(kw)
        elif project_name in kw:
            matched.append(kw)
    return matched


def infer_university_type(name):
    """Infer university type from its name"""
    if any(k in name for k in ["体育学院", "体育大学", "体育职业"]):
        return "体育类"
    if any(k in name for k in ["师范大学", "师范学院", "师范高等专科"]):
        return "师范类"
    if any(k in name for k in ["民族大学", "民族学院"]):
        return "民族类"
    if any(k in name for k in ["大学", "学院"]):
        return "综合类"
    return "综合类"


def infer_location(region):
    """Obtain location based on the region field"""
    # The region field is already a province name, use it directly
    return region


# ========================== Main Program Entry ==========================
def main():
    print("=" * 70)
    print("Ministry of Education Traditional Culture Inheritance Base Intangible Heritage Sports Course Data Crawler")
    print("Data source: three batches of 'Outstanding Traditional Chinese Culture Inheritance Bases in Regular Higher Education Institutions' published on the MOE official website")
    print("=" * 70)

    try:
        # ---------- Step 1: Load matching keywords ----------
        keywords = read_keywords()
        print(f"Keyword preview (first 10 items): {keywords[:10]}")

        # ---------- Step 2: Merge three batches of base data ----------
        all_bases = []
        # Add batch label for each record
        for item in BATCH1_BASE:
            item_copy = item.copy()
            item_copy["批次"] = "第一批（2018年）"
            all_bases.append(item_copy)
        for item in BATCH2_BASE:
            item_copy = item.copy()
            item_copy["批次"] = "第二批（2019年）"
            all_bases.append(item_copy)
        for item in BATCH3_BASE:
            item_copy = item.copy()
            item_copy["批次"] = "第三批（2020年）"
            all_bases.append(item_copy)

        print(f"\n📊 Total {len(all_bases)} base records loaded (sum of three batches)")

        # ---------- Step 3: Filter traditional sports bases and match keywords ----------
        print("\n" + "=" * 70)
        print("🔍 Start matching intangible heritage sports projects")
        print("=" * 70)

        # Define traditional sports related keywords (used for preliminary filtering)
        sports_keywords = ["武术", "太极", "射艺", "龙舟", "拳", "跤", "摔", "剑", "棍", "枪", "刀",
                           "舞狮", "八段锦", "五禽戏", "民族传统体育", "体育", "健身气功", "蹴鞠", "马球",
                           "摔跤", "搏克", "秋千", "跳板", "赛马", "射箭", "陀螺", "珍珠球"]

        results = []
        processed_count = 0
        matched_count = 0

        for idx, base in enumerate(all_bases, 1):
            school = base.get("学校名称", "")
            region = base.get("地区", "")
            project = base.get("传承项目", "")
            batch = base.get("批次", "")

            # Determine whether it is a traditional sports base (judged by inheritance project name)
            is_sports = False
            for sk in sports_keywords:
                if sk in project:
                    is_sports = True
                    break

            if not is_sports:
                continue

            processed_count += 1

            # Match user keywords
            matched_kws = match_keyword(project, keywords)

            if matched_kws:
                matched_count += 1
                print(f"  ✅ {school}：{project} → Matched: {', '.join(matched_kws)}")

                for kw in matched_kws:
                    results.append({
                        "对应非遗体育项目名称": kw,
                        "院校名称": school,
                        "院校所在地": infer_location(region),
                        "院校类型": infer_university_type(school),
                        "非遗传统体育课程名称": project,  # The inheritance project is the course name
                        "课程属性": "公共选修课",  # MOE requires opening elective courses for the whole university
                        "课程等级": "本科",
                        "课程简介": f"教育部中华优秀传统文化传承基地，传承项目为「{project}」，"
                                    f"面向全校学生开设2学分选修课程。{batch}认定。",
                        "数据来源渠道": "教育部传统文化传承基地公示",
                        "备注": "国家级中华优秀传统文化传承基地"
                    })
            else:
                # Traditional sports base but no keyword matched; still record it with empty project name
                print(f"  ⚠️ {school}：{project}（No keyword matched）")
                results.append({
                    "对应非遗体育项目名称": "",
                    "院校名称": school,
                    "院校所在地": infer_location(region),
                    "院校类型": infer_university_type(school),
                    "非遗传统体育课程名称": project,
                    "课程属性": "公共选修课",
                    "课程等级": "本科",
                    "课程简介": f"教育部中华优秀传统文化传承基地，传承项目为「{project}」。{batch}认定。",
                    "数据来源渠道": "教育部传统文化传承基地公示",
                    "备注": "国家级中华优秀传统文化传承基地（No keyword matched）"
                })

        print(f"\n📊 Traditional sports bases: {processed_count}")
        print(f"📊 Bases matched with keywords: {matched_count}")

        # ---------- Step 4: Deduplicate ----------
        if results:
            df_results = pd.DataFrame(results)
            # Deduplicate by university name + intangible heritage sports course name
            df_results = df_results.drop_duplicates(
                subset=["院校名称", "非遗传统体育课程名称"],
                keep="first"
            )
        else:
            df_results = pd.DataFrame(columns=[
                "对应非遗体育项目名称", "院校名称", "院校所在地", "院校类型",
                "非遗传统体育课程名称", "课程属性", "课程等级", "课程简介",
                "数据来源渠道", "备注"
            ])

        # ---------- Step 5: Export to excel ----------
        df_results.to_excel(OUTPUT_FILE, index=False)

        # ---------- Step 6: Print statistical report ----------
        print("\n" + "=" * 70)
        print("📊 Crawling‑task Completion Statistics")
        print("=" * 70)
        print(f"Total matching keywords: {len(keywords)}")
        print(f"Total base records across three batches: {len(all_bases)}")
        print(f"Traditional sports base count: {processed_count}")
        print(f"Bases matched successfully: {matched_count}")
        print(f"Exported records (including unmatched): {len(results)}")
        print(f"Deduplicated records: {len(df_results)}")
        print(f"\n✅ Output file: {OUTPUT_FILE}")
        print("=" * 70)

        # Preview result data
        if not df_results.empty:
            print("\n📖 Result preview (first 10 entries, matched only):")
            print("-" * 70)
            matched_df = df_results[df_results["对应非遗体育项目名称"] != ""]
            if not matched_df.empty:
                preview_cols = ["对应非遗体育项目名称", "院校名称", "非遗传统体育课程名称"]
                print(matched_df[preview_cols].head(10).to_string(index=False))
            else:
                print("No keyword‑matched records yet")
            print("-" * 70)

            # Count matched universities per project
            print("\n📊 Number of matched universities per intangible heritage sports project:")
            matched_df = df_results[df_results["对应非遗体育项目名称"] != ""]
            if not matched_df.empty:
                course_count = matched_df.groupby("对应非遗体育项目名称").size().sort_values(ascending=False)
                for project, count in course_count.items():
                    print(f"  {project}: {count} universities")
        else:
            print("\n⚠️ No records generated")

        # ---------- Step 7: Data source description ----------
        print("\n" + "=" * 70)
        print("📌 Data Source Description")
        print("=" * 70)
        print("The data comes from the three batches of 'Outstanding Traditional Chinese Culture Inheritance Bases in Regular Higher Education Institutions' published on the MOE official website:")
        print("  - Batch 1 (2018): Document 教体艺厅函〔2018〕85号, 55 bases")
        print("  - Batch 2 (2019): Document 教体艺厅函〔2019〕53号, 25 bases")
        print("  - Batch 3 (2020): Document 教体艺厅函〔2020〕26号, 26 bases")
        print("Total 106 bases.")
        print("\n💡 Notes:")
        print("  - The '课程属性' field is uniformly labeled as '公共选修课' according to MOE requirements")
        print("  - MOE requires each base to open elective courses based on its inheritance project for all students")
        print("  - The courses are incorporated into talent training plans and teaching schedules, with credit management and 2 credits")
        print("  - All bases are 'National Outstanding Traditional Chinese Culture Inheritance Bases'")
        print("=" * 70)

    except FileNotFoundError as e:
        print(f"❌ File‑related error: {e}")
    except ValueError as e:
        print(f"❌ Data‑validation error: {e}")
    except Exception as e:
        print(f"❌ Unexpected runtime error: {e}")
        import traceback
        traceback.print_exc()


if __name__ == "__main__":
    main()
